agora inbox for pgsql-hackers@postgresql.org  
help / color / mirror / Atom feed
[PATCH v44 07/10] Split cluster.h to create repack_internal.h
324+ messages / 1 participants
[nested] [flat]

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread

* [PATCH v56 1/3] Allow logical replication snapshots to be database-specific
@ 2026-04-06 22:07 Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 324+ messages in thread

From: Álvaro Herrera @ 2026-04-06 22:07 UTC (permalink / raw)

By default, the logical decoding assumes access to shared catalogs, so
the snapshot builder needs to consider cluster-wide XIDs during startup.
That in turn means that, if any transaction is already running (and has
XID assigned), the snapshot builder needs to wait for its completion, as
it does not know if that transaction performed catalog changes earlier.

A possible problem with this concept is that if REPACK (CONCURRENTLY) is
running in some database, backends running the same command in other
databases get stuck until the first one has committed. Thus only a
single backend in the cluster can run REPACK (CONCURRENTLY) at any time.
Likewise, REPACK (CONCURRENTLY) can block walsenders starting on behalf
of subscriptions throughout the cluster.

This patch adds a new option to logical replication output plugin, to
declare that it does not use shared catalogs (i.e. catalogs that can be
changed by transactions running in other databases in the cluster). In
that case, no snapshot the backend will use during the decoding needs to
contain information about transactions running in other databases. Thus
the snapshot builder only needs to wait for completion of transactions
in the current database.

Currently we only use this option in the REPACK background worker. It
could possibly be used in the plugin for logical replication too,
however that would need thorough analysis of that plugin.

Bump WAL version number, due to a new field in xl_running_xacts.

Author: Antonin Houska <ah@cybertec.at>
Reviewed-by: Álvaro Herrera <alvherre@kurilemu.de>
Discussion: https://postgr.es/m/90475.1775218118@localhost
---
 contrib/pg_visibility/pg_visibility.c       |  4 +-
 doc/src/sgml/logicaldecoding.sgml           |  4 ++
 src/backend/access/index/genam.c            |  8 +++
 src/backend/access/rmgrdesc/standbydesc.c   |  2 +
 src/backend/access/transam/xlog.c           |  2 +-
 src/backend/access/transam/xlogfuncs.c      |  2 +-
 src/backend/postmaster/bgwriter.c           |  2 +-
 src/backend/replication/logical/decode.c    | 17 +++++-
 src/backend/replication/logical/logical.c   |  3 +
 src/backend/replication/logical/snapbuild.c | 63 ++++++++++++++++++++-
 src/backend/replication/pgrepack/pgrepack.c |  7 +++
 src/backend/replication/slot.c              |  2 +-
 src/backend/storage/ipc/procarray.c         | 23 +++++++-
 src/backend/storage/ipc/standby.c           | 24 +++++++-
 src/include/access/genam.h                  |  1 +
 src/include/access/xlog_internal.h          |  2 +-
 src/include/miscadmin.h                     |  9 +++
 src/include/replication/output_plugin.h     |  1 +
 src/include/replication/snapbuild.h         |  3 +-
 src/include/storage/procarray.h             |  2 +-
 src/include/storage/standby.h               |  3 +-
 src/include/storage/standbydefs.h           |  1 +
 22 files changed, 167 insertions(+), 18 deletions(-)

diff --git a/contrib/pg_visibility/pg_visibility.c b/contrib/pg_visibility/pg_visibility.c
index dfab0b64cf5..d564bd2a00c 100644
--- a/contrib/pg_visibility/pg_visibility.c
+++ b/contrib/pg_visibility/pg_visibility.c
@@ -621,7 +621,7 @@ GetStrictOldestNonRemovableTransactionId(Relation rel)
 	else if (rel == NULL || rel->rd_rel->relisshared)
 	{
 		/* Shared relation: take into account all running xids */
-		runningTransactions = GetRunningTransactionData();
+		runningTransactions = GetRunningTransactionData(InvalidOid);
 		LWLockRelease(ProcArrayLock);
 		LWLockRelease(XidGenLock);
 		return runningTransactions->oldestRunningXid;
@@ -632,7 +632,7 @@ GetStrictOldestNonRemovableTransactionId(Relation rel)
 		 * Normal relation: take into account xids running within the current
 		 * database
 		 */
-		runningTransactions = GetRunningTransactionData();
+		runningTransactions = GetRunningTransactionData(InvalidOid);
 		LWLockRelease(ProcArrayLock);
 		LWLockRelease(XidGenLock);
 		return runningTransactions->oldestDatabaseRunningXid;
diff --git a/doc/src/sgml/logicaldecoding.sgml b/doc/src/sgml/logicaldecoding.sgml
index 6dc49108997..9b1d68d0de6 100644
--- a/doc/src/sgml/logicaldecoding.sgml
+++ b/doc/src/sgml/logicaldecoding.sgml
@@ -959,6 +959,7 @@ typedef struct OutputPluginOptions
 {
     OutputPluginOutputType output_type;
     bool        receive_rewrites;
+    bool        need_shared_catalogs;
 } OutputPluginOptions;
 </programlisting>
       <literal>output_type</literal> has to either be set to
@@ -969,6 +970,9 @@ typedef struct OutputPluginOptions
       also be called for changes made by heap rewrites during certain DDL
       operations.  These are of interest to plugins that handle DDL
       replication, but they require special handling.
+      <literal>need_shared_catalogs</literal> can be set to false if you are
+      certain the plugin functions do not access shared system catalogs.
+      Doing so can speed up creation of replication slots that use this plugin.
      </para>
 
      <para>
diff --git a/src/backend/access/index/genam.c b/src/backend/access/index/genam.c
index 1408989c568..97d44b84622 100644
--- a/src/backend/access/index/genam.c
+++ b/src/backend/access/index/genam.c
@@ -394,6 +394,14 @@ systable_beginscan(Relation heapRelation,
 	SysScanDesc sysscan;
 	Relation	irel;
 
+	/*
+	 * If this backend promised that it won't access shared catalogs during
+	 * logical decoding, this it the right place to verify.
+	 */
+	Assert(!HistoricSnapshotActive() ||
+		   accessSharedCatalogsInDecoding ||
+		   !heapRelation->rd_rel->relisshared);
+
 	if (indexOK &&
 		!IgnoreSystemIndexes &&
 		!ReindexIsProcessingIndex(indexId))
diff --git a/src/backend/access/rmgrdesc/standbydesc.c b/src/backend/access/rmgrdesc/standbydesc.c
index 0a291354ae2..685d1bdb024 100644
--- a/src/backend/access/rmgrdesc/standbydesc.c
+++ b/src/backend/access/rmgrdesc/standbydesc.c
@@ -41,6 +41,8 @@ standby_desc_running_xacts(StringInfo buf, xl_running_xacts *xlrec)
 		for (i = 0; i < xlrec->subxcnt; i++)
 			appendStringInfo(buf, " %u", xlrec->xids[xlrec->xcnt + i]);
 	}
+
+	appendStringInfo(buf, "; dbid: %u", xlrec->dbid);
 }
 
 void
diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c
index b82af9a85c0..3f08a832ca6 100644
--- a/src/backend/access/transam/xlog.c
+++ b/src/backend/access/transam/xlog.c
@@ -7735,7 +7735,7 @@ CreateCheckPoint(int flags)
 	 * recovery we don't need to write running xact data.
 	 */
 	if (!shutdown && XLogStandbyInfoActive())
-		LogStandbySnapshot();
+		LogStandbySnapshot(InvalidOid);
 
 	START_CRIT_SECTION();
 
diff --git a/src/backend/access/transam/xlogfuncs.c b/src/backend/access/transam/xlogfuncs.c
index 65bbaeda59c..0f5979691e6 100644
--- a/src/backend/access/transam/xlogfuncs.c
+++ b/src/backend/access/transam/xlogfuncs.c
@@ -245,7 +245,7 @@ pg_log_standby_snapshot(PG_FUNCTION_ARGS)
 				(errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE),
 				 errmsg("pg_log_standby_snapshot() can only be used if \"wal_level\" >= \"replica\"")));
 
-	recptr = LogStandbySnapshot();
+	recptr = LogStandbySnapshot(InvalidOid);
 
 	/*
 	 * As a convenience, return the WAL location of the last inserted record
diff --git a/src/backend/postmaster/bgwriter.c b/src/backend/postmaster/bgwriter.c
index 1d8947774a9..a30de4262eb 100644
--- a/src/backend/postmaster/bgwriter.c
+++ b/src/backend/postmaster/bgwriter.c
@@ -289,7 +289,7 @@ BackgroundWriterMain(const void *startup_data, size_t startup_data_len)
 			if (now >= timeout &&
 				last_snapshot_lsn <= GetLastImportantRecPtr())
 			{
-				last_snapshot_lsn = LogStandbySnapshot();
+				last_snapshot_lsn = LogStandbySnapshot(InvalidOid);
 				last_snapshot_ts = now;
 			}
 		}
diff --git a/src/backend/replication/logical/decode.c b/src/backend/replication/logical/decode.c
index c9fea8cad28..38c5a4f5540 100644
--- a/src/backend/replication/logical/decode.c
+++ b/src/backend/replication/logical/decode.c
@@ -382,7 +382,16 @@ standby_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf)
 			{
 				xl_running_xacts *running = (xl_running_xacts *) XLogRecGetData(r);
 
-				SnapBuildProcessRunningXacts(builder, buf->origptr, running);
+				/*
+				 * Update this decoder's idea of transactions currently
+				 * running.  In doing so we will determine whether we have
+				 * reached consistent status.
+				 *
+				 * If the output plugin doesn't need access to shared
+				 * catalogs, we can ignore transactions in other databases.
+				 */
+				SnapBuildProcessRunningXacts(builder, buf->origptr, running,
+											 !ctx->options.need_shared_catalogs);
 
 				/*
 				 * Abort all transactions that we keep track of, that are
@@ -392,8 +401,12 @@ standby_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf)
 				 * all running transactions which includes prepared ones,
 				 * while shutdown checkpoints just know that no non-prepared
 				 * transactions are in progress.
+				 *
+				 * The database-specific records might work here too, but it's
+				 * not their purpose.
 				 */
-				ReorderBufferAbortOld(ctx->reorder, running->oldestRunningXid);
+				if (!OidIsValid(running->dbid))
+					ReorderBufferAbortOld(ctx->reorder, running->oldestRunningXid);
 			}
 			break;
 		case XLOG_STANDBY_LOCK:
diff --git a/src/backend/replication/logical/logical.c b/src/backend/replication/logical/logical.c
index f20d0c542f3..8ceaf64d164 100644
--- a/src/backend/replication/logical/logical.c
+++ b/src/backend/replication/logical/logical.c
@@ -285,6 +285,9 @@ StartupDecodingContext(List *output_plugin_options,
 	ctx->write = do_write;
 	ctx->update_progress = update_progress;
 
+	/* Assume shared catalog access. The startup callback can change it. */
+	ctx->options.need_shared_catalogs = true;
+
 	ctx->output_plugin_options = output_plugin_options;
 
 	ctx->fast_forward = fast_forward;
diff --git a/src/backend/replication/logical/snapbuild.c b/src/backend/replication/logical/snapbuild.c
index b4269a3b102..ffb4ab2cf2a 100644
--- a/src/backend/replication/logical/snapbuild.c
+++ b/src/backend/replication/logical/snapbuild.c
@@ -154,6 +154,14 @@
 static ResourceOwner SavedResourceOwnerDuringExport = NULL;
 static bool ExportInProgress = false;
 
+/*
+ * If a backend is going to do logical decoding and the output plugin does
+ * not need to access shared catalogs, setting this variable to false can make
+ * the decoding startup faster. In particular, the backend will not need to
+ * wait for completion of already running transactions in other databases.
+ */
+bool		accessSharedCatalogsInDecoding = true;
+
 /* ->committed and ->catchange manipulation */
 static void SnapBuildPurgeOlderTxn(SnapBuild *builder);
 
@@ -170,7 +178,8 @@ static inline bool SnapBuildXidHasCatalogChanges(SnapBuild *builder, Transaction
 												 uint32 xinfo);
 
 /* xlog reading helper functions for SnapBuildProcessRunningXacts */
-static bool SnapBuildFindSnapshot(SnapBuild *builder, XLogRecPtr lsn, xl_running_xacts *running);
+static bool SnapBuildFindSnapshot(SnapBuild *builder, XLogRecPtr lsn,
+								  xl_running_xacts *running);
 static void SnapBuildWaitSnapshot(xl_running_xacts *running, TransactionId cutoff);
 
 /* serialization functions */
@@ -226,6 +235,9 @@ AllocateSnapshotBuilder(ReorderBuffer *reorder,
 
 	MemoryContextSwitchTo(oldcontext);
 
+	/* The default is that shared catalog are used. */
+	accessSharedCatalogsInDecoding = true;
+
 	return builder;
 }
 
@@ -244,6 +256,9 @@ FreeSnapshotBuilder(SnapBuild *builder)
 		builder->snapshot = NULL;
 	}
 
+	/* The default is that shared catalog are used. */
+	accessSharedCatalogsInDecoding = true;
+
 	/* other resources are deallocated via memory context reset */
 	MemoryContextDelete(context);
 }
@@ -1136,7 +1151,8 @@ SnapBuildXidHasCatalogChanges(SnapBuild *builder, TransactionId xid,
  * anymore.
  */
 void
-SnapBuildProcessRunningXacts(SnapBuild *builder, XLogRecPtr lsn, xl_running_xacts *running)
+SnapBuildProcessRunningXacts(SnapBuild *builder, XLogRecPtr lsn, xl_running_xacts *running,
+							 bool db_specific)
 {
 	ReorderBufferTXN *txn;
 	TransactionId xmin;
@@ -1148,6 +1164,33 @@ SnapBuildProcessRunningXacts(SnapBuild *builder, XLogRecPtr lsn, xl_running_xact
 	 */
 	if (builder->state < SNAPBUILD_CONSISTENT)
 	{
+		/*
+		 * To reduce the potential for unnecessarily waiting for completion of
+		 * unrelated transactions, the caller can declare that only
+		 * transactions of the current database are relevant at this stage.
+		 */
+		if (db_specific)
+		{
+			/*
+			 * If we must only keep track of transactions running in the
+			 * current database, we need transaction info from exactly that
+			 * database.
+			 */
+			if (running->dbid != MyDatabaseId)
+			{
+				LogStandbySnapshot(MyDatabaseId);
+
+				return;
+			}
+
+			/*
+			 * We'd better be able to check during scan if the plugin does not
+			 * lie.
+			 */
+			if (accessSharedCatalogsInDecoding)
+				accessSharedCatalogsInDecoding = false;
+		}
+
 		/* returns false if there's no point in performing cleanup just yet */
 		if (!SnapBuildFindSnapshot(builder, lsn, running))
 			return;
@@ -1155,6 +1198,16 @@ SnapBuildProcessRunningXacts(SnapBuild *builder, XLogRecPtr lsn, xl_running_xact
 	else
 		SnapBuildSerialize(builder, lsn);
 
+	/*
+	 * Database specific transaction info may exist to reach CONSISTENT state
+	 * faster, however the code below makes no use of it. Moreover, such
+	 * record might cause problems because the following normal (cluster-wide)
+	 * record can have lower value of oldestRunningXid. In that case, let's
+	 * wait with the cleanup for the next regular cluster-wide record.
+	 */
+	if (OidIsValid(running->dbid))
+		return;
+
 	/*
 	 * Update range of interesting xids based on the running xacts
 	 * information. We don't increase ->xmax using it, because once we are in
@@ -1465,7 +1518,11 @@ SnapBuildWaitSnapshot(xl_running_xacts *running, TransactionId cutoff)
 	 */
 	if (!RecoveryInProgress())
 	{
-		LogStandbySnapshot();
+		/*
+		 * If the last transaction info was about specific database, so needs
+		 * to be the next one - at least until we're in the CONSISTENT state.
+		 */
+		LogStandbySnapshot(running->dbid);
 	}
 }
 
diff --git a/src/backend/replication/pgrepack/pgrepack.c b/src/backend/replication/pgrepack/pgrepack.c
index 457b6518a8e..4da66aa0f44 100644
--- a/src/backend/replication/pgrepack/pgrepack.c
+++ b/src/backend/replication/pgrepack/pgrepack.c
@@ -52,6 +52,13 @@ repack_startup(LogicalDecodingContext *ctx, OutputPluginOptions *opt,
 	/* Probably unnecessary, as we don't use the SQL interface ... */
 	opt->output_type = OUTPUT_PLUGIN_BINARY_OUTPUT;
 
+	/*
+	 * REPACK doesn't need access to shared catalogs, so we can speed up the
+	 * historic snapshot creation by setting this flag.  We'll only have to
+	 * wait for transactions in our database.
+	 */
+	opt->need_shared_catalogs = false;
+
 	if (ctx->output_plugin_options != NIL)
 	{
 		ereport(ERROR,
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 21a213a0ebf..a1f37e59dbc 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -1756,7 +1756,7 @@ ReplicationSlotReserveWal(void)
 		XLogRecPtr	flushptr;
 
 		/* make sure we have enough information to start */
-		flushptr = LogStandbySnapshot();
+		flushptr = LogStandbySnapshot(InvalidOid);
 
 		/* and make sure it's fsynced to disk */
 		XLogFlush(flushptr);
diff --git a/src/backend/storage/ipc/procarray.c b/src/backend/storage/ipc/procarray.c
index f540bb6b23f..9299bcebbda 100644
--- a/src/backend/storage/ipc/procarray.c
+++ b/src/backend/storage/ipc/procarray.c
@@ -2623,9 +2623,11 @@ ProcArrayInstallRestoredXmin(TransactionId xmin, PGPROC *proc)
  *
  * Note that if any transaction has overflowed its cached subtransactions
  * then there is no real need include any subtransactions.
+ *
+ * If 'dbid' is valid, only gather transactions running in that database.
  */
 RunningTransactions
-GetRunningTransactionData(void)
+GetRunningTransactionData(Oid dbid)
 {
 	/* result workspace */
 	static RunningTransactionsData CurrentRunningXactsData;
@@ -2700,6 +2702,18 @@ GetRunningTransactionData(void)
 		if (!TransactionIdIsValid(xid))
 			continue;
 
+		/*
+		 * Filter by database OID if requested.
+		 */
+		if (OidIsValid(dbid))
+		{
+			int			pgprocno = arrayP->pgprocnos[index];
+			PGPROC	   *proc = &allProcs[pgprocno];
+
+			if (proc->databaseId != dbid)
+				continue;
+		}
+
 		/*
 		 * Be careful not to exclude any xids before calculating the values of
 		 * oldestRunningXid and suboverflowed, since these are used to clean
@@ -2750,6 +2764,12 @@ GetRunningTransactionData(void)
 			PGPROC	   *proc = &allProcs[pgprocno];
 			int			nsubxids;
 
+			/*
+			 * Filter by database OID if requested.
+			 */
+			if (OidIsValid(dbid) && proc->databaseId != dbid)
+				continue;
+
 			/*
 			 * Save subtransaction XIDs. Other backends can't add or remove
 			 * entries while we're holding XidGenLock.
@@ -2783,6 +2803,7 @@ GetRunningTransactionData(void)
 	 * increases if slots do.
 	 */
 
+	CurrentRunningXacts->dbid = dbid;
 	CurrentRunningXacts->xcnt = count - subcount;
 	CurrentRunningXacts->subxcnt = subcount;
 	CurrentRunningXacts->subxid_status = suboverflowed ? SUBXIDS_IN_SUBTRANS : SUBXIDS_IN_ARRAY;
diff --git a/src/backend/storage/ipc/standby.c b/src/backend/storage/ipc/standby.c
index de9092fdf5b..29af7733948 100644
--- a/src/backend/storage/ipc/standby.c
+++ b/src/backend/storage/ipc/standby.c
@@ -1188,6 +1188,14 @@ standby_redo(XLogReaderState *record)
 		xl_running_xacts *xlrec = (xl_running_xacts *) XLogRecGetData(record);
 		RunningTransactionsData running;
 
+		/*
+		 * Records issued for specific database are not suitable for physical
+		 * replication because that affects the whole cluster. In particular,
+		 * the list of XID is probably incomplete here.
+		 */
+		if (OidIsValid(xlrec->dbid))
+			return;
+
 		running.xcnt = xlrec->xcnt;
 		running.subxcnt = xlrec->subxcnt;
 		running.subxid_status = xlrec->subxid_overflow ? SUBXIDS_MISSING : SUBXIDS_IN_ARRAY;
@@ -1277,11 +1285,22 @@ standby_redo(XLogReaderState *record)
  * as there's no independent knob to just enable logical decoding. For
  * details of how this is used, check snapbuild.c's introductory comment.
  *
+ * If 'dbid' is valid, only gather transactions running in that
+ * database. snapbuild.c can use such running xacts information for faster
+ * startup, but it still needs normal (cluster-wide) during the actual
+ * decoding - see standby_decode() and SnapBuildProcessRunningXacts() for
+ * details. Other processes (e.g. checkpointer) issue the cluster-wide records
+ * whether logical decoding is active or not.
+ *
+ * Please be careful about using this argument for other purposes. In
+ * particular, physical replication *must* ignore the database-specific
+ * records, exactly because they do not cover the whole cluster - see
+ * standby_redo().
  *
  * Returns the RecPtr of the last inserted record.
  */
 XLogRecPtr
-LogStandbySnapshot(void)
+LogStandbySnapshot(Oid dbid)
 {
 	XLogRecPtr	recptr;
 	RunningTransactions running;
@@ -1314,7 +1333,7 @@ LogStandbySnapshot(void)
 	 * Log details of all in-progress transactions. This should be the last
 	 * record we write, because standby will open up when it sees this.
 	 */
-	running = GetRunningTransactionData();
+	running = GetRunningTransactionData(dbid);
 
 	/*
 	 * GetRunningTransactionData() acquired ProcArrayLock, we must release it.
@@ -1358,6 +1377,7 @@ LogCurrentRunningXacts(RunningTransactions CurrRunningXacts)
 	xl_running_xacts xlrec;
 	XLogRecPtr	recptr;
 
+	xlrec.dbid = CurrRunningXacts->dbid;
 	xlrec.xcnt = CurrRunningXacts->xcnt;
 	xlrec.subxcnt = CurrRunningXacts->subxcnt;
 	xlrec.subxid_overflow = (CurrRunningXacts->subxid_status != SUBXIDS_IN_ARRAY);
diff --git a/src/include/access/genam.h b/src/include/access/genam.h
index b69320a7fc8..de55821e414 100644
--- a/src/include/access/genam.h
+++ b/src/include/access/genam.h
@@ -136,6 +136,7 @@ typedef struct IndexOrderByDistance
 	bool		isnull;
 } IndexOrderByDistance;
 
+
 /*
  * generalized index_ interface routines (in indexam.c)
  */
diff --git a/src/include/access/xlog_internal.h b/src/include/access/xlog_internal.h
index 10c18d39ff8..13ae3ad4fbb 100644
--- a/src/include/access/xlog_internal.h
+++ b/src/include/access/xlog_internal.h
@@ -32,7 +32,7 @@
 /*
  * Each page of XLOG file has a header like this:
  */
-#define XLOG_PAGE_MAGIC 0xD11E	/* can be used as WAL version indicator */
+#define XLOG_PAGE_MAGIC 0xD11F	/* can be used as WAL version indicator */
 
 typedef struct XLogPageHeaderData
 {
diff --git a/src/include/miscadmin.h b/src/include/miscadmin.h
index 2e10e3c814d..93b7816c09c 100644
--- a/src/include/miscadmin.h
+++ b/src/include/miscadmin.h
@@ -310,6 +310,15 @@ extern void PreventCommandIfReadOnly(const char *cmdname);
 extern void PreventCommandIfParallelMode(const char *cmdname);
 extern void PreventCommandDuringRecovery(const char *cmdname);
 
+/* in replication/snapbuild.c */
+
+/*
+ * Keep track of whether logical decoding in this backend promised not to
+ * access shared catalogs, as a safety check.  This is checked by genam.c when
+ * a catalog scan takes place to verify that no shared catalogs are accessed.
+ */
+extern bool accessSharedCatalogsInDecoding;
+
 /*****************************************************************************
  *	  pdir.h --																 *
  *			POSTGRES directory path definitions.                             *
diff --git a/src/include/replication/output_plugin.h b/src/include/replication/output_plugin.h
index 842fcde67f9..917f3cff232 100644
--- a/src/include/replication/output_plugin.h
+++ b/src/include/replication/output_plugin.h
@@ -27,6 +27,7 @@ typedef struct OutputPluginOptions
 {
 	OutputPluginOutputType output_type;
 	bool		receive_rewrites;
+	bool		need_shared_catalogs;
 } OutputPluginOptions;
 
 /*
diff --git a/src/include/replication/snapbuild.h b/src/include/replication/snapbuild.h
index a22a83a2f23..d02530a912a 100644
--- a/src/include/replication/snapbuild.h
+++ b/src/include/replication/snapbuild.h
@@ -92,7 +92,8 @@ extern void SnapBuildProcessNewCid(SnapBuild *builder, TransactionId xid,
 								   XLogRecPtr lsn,
 								   xl_heap_new_cid *xlrec);
 extern void SnapBuildProcessRunningXacts(SnapBuild *builder, XLogRecPtr lsn,
-										 xl_running_xacts *running);
+										 xl_running_xacts *running,
+										 bool db_specific);
 extern void SnapBuildSerializationPoint(SnapBuild *builder, XLogRecPtr lsn);
 
 extern bool SnapBuildSnapshotExists(XLogRecPtr lsn);
diff --git a/src/include/storage/procarray.h b/src/include/storage/procarray.h
index d718a5b542f..ec89c448220 100644
--- a/src/include/storage/procarray.h
+++ b/src/include/storage/procarray.h
@@ -47,7 +47,7 @@ extern bool ProcArrayInstallImportedXmin(TransactionId xmin,
 										 VirtualTransactionId *sourcevxid);
 extern bool ProcArrayInstallRestoredXmin(TransactionId xmin, PGPROC *proc);
 
-extern RunningTransactions GetRunningTransactionData(void);
+extern RunningTransactions GetRunningTransactionData(Oid dbid);
 
 extern bool TransactionIdIsInProgress(TransactionId xid);
 extern TransactionId GetOldestNonRemovableTransactionId(Relation rel);
diff --git a/src/include/storage/standby.h b/src/include/storage/standby.h
index 6a314c693cd..8715c08e94f 100644
--- a/src/include/storage/standby.h
+++ b/src/include/storage/standby.h
@@ -126,6 +126,7 @@ typedef enum
 
 typedef struct RunningTransactionsData
 {
+	Oid			dbid;			/* only track xacts in this database */
 	int			xcnt;			/* # of xact ids in xids[] */
 	int			subxcnt;		/* # of subxact ids in xids[] */
 	subxids_array_status subxid_status;
@@ -143,7 +144,7 @@ typedef RunningTransactionsData *RunningTransactions;
 extern void LogAccessExclusiveLock(Oid dbOid, Oid relOid);
 extern void LogAccessExclusiveLockPrepare(void);
 
-extern XLogRecPtr LogStandbySnapshot(void);
+extern XLogRecPtr LogStandbySnapshot(Oid dbid);
 extern void LogStandbyInvalidations(int nmsgs, SharedInvalidationMessage *msgs,
 									bool relcacheInitFileInval);
 
diff --git a/src/include/storage/standbydefs.h b/src/include/storage/standbydefs.h
index 231d251fd51..e75b7078766 100644
--- a/src/include/storage/standbydefs.h
+++ b/src/include/storage/standbydefs.h
@@ -46,6 +46,7 @@ typedef struct xl_standby_locks
  */
 typedef struct xl_running_xacts
 {
+	Oid			dbid;			/* only track xacts in this database */
 	int			xcnt;			/* # of xact ids in xids[] */
 	int			subxcnt;		/* # of subxact ids in xids[] */
 	bool		subxid_overflow;	/* snapshot overflowed, subxids missing */
-- 
2.47.3


--qs77q6fpwolne4ds
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v56-0002-Reserve-replication-slots-specifically-for-REPAC.patch"



^ permalink  raw  reply  [nested|flat] 324+ messages in thread


end of thread, other threads:[~2026-04-06 22:07 UTC | newest]

Thread overview: 324+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-04-06 22:07 [PATCH v56 1/3] Allow logical replication snapshots to be database-specific Álvaro Herrera <alvherre@kurilemu.de>

This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox